You are an expert CUDA programmer. Your task is to replace a PyTorch operator with a highly optimized and numerically stable custom CUDA kernel.

Here's an example to show you the syntax of inline embedding custom CUDA operators in PyTorch. The example given architecture is a simple ReLU:

python
import torch
import torch.nn as nn
import torch.nn.functional as F

class Model(nn.Module):
def init(self) -> None:
super().init()

def forward(self, x):
    return torch.relu(x)
def get_inputs():
x = torch.randn(1, 128).cuda()
return [x]

def get_init_inputs():
return []



The example new architecture with a custom CUDA kernel looks like this:

python
import torch
from torch.utils.cpp_extension import load_inline

relu_source = """
#include <torch/extension.h>
#include <cuda_runtime.h>

global void relu_kernel(const float* x, float* y, int size) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if (idx < size) {
y[idx] = fmaxf(x[idx], 0.f);
}
}

torch::Tensor relu_cuda(torch::Tensor x) {
auto size = x.numel();
auto y = torch::empty_like(x);
const int block_size = 256;
int num_blocks = (size + block_size - 1) / block_size;
relu_kernel<<<num_blocks, block_size>>>(x.data_ptr<float>(), y.data_ptr<float>(), size);
return y;
}
"""

relu_cpp_source = """
torch::Tensor relu_cuda(torch::Tensor x);
"""

Compile the inline CUDA code
relu = load_inline(
name=“relu”,
cpp_sources=relu_cpp_source,
cuda_sources=relu_source,
functions=[“relu_cuda”],
verbose=True
)

class ModelNew(torch.nn.Module):
def init(self):
super(ModelNew, self).init()
self.relu = relu # The module containing the kernel

def forward(self, x):
    return self.relu.relu_cuda(x)
def get_inputs():
x = torch.randn(1, 128).cuda()
return [x]

def get_init_inputs():
return []



---

Now, you are given the following PyTorch architecture to accelerate:

python
import torch
import torch.nn as nn

class Model(nn.Module):
“”"
Manhattan + Sqrt fusion implementation.
Computes the Manhattan distance (L1 distance) between two sets of vectors and then applies a square root.
“”"
def init(self):
super(Model, self).init()

def forward(self, x: torch.Tensor, y: torch.Tensor) -> torch.Tensor:
    """
    Compute Manhattan distance between x and y, then apply sqrt.

    Args:
        x (torch.Tensor): First set of vectors [batch_size, feature_dim]
        y (torch.Tensor): Second set of vectors [batch_size, feature_dim]

    Returns:
        torch.Tensor: Square root of Manhattan distances [batch_size]
    """
    # Input validation
    if x.shape != y.shape:
        raise ValueError(f"Input tensors must have the same shape, got {x.shape} and {y.shape}")
    
    if x.dim() != 2:
        raise ValueError(f"Input tensors must be 2D, got {x.dim()}D")
    
    # Step 1: Compute Manhattan distance: Σ|x_i - y_i|
    manhattan_dist = torch.sum(torch.abs(x - y), dim=1)
    
    # Step 2: Apply square root
    sqrt_manhattan_dist = torch.sqrt(manhattan_dist)
    
    return sqrt_manhattan_dist
batch_size = 256
feature_dim = 512

def get_inputs():
# Generate two sets of vectors
x = torch.randn(batch_size, feature_dim)
y = torch.randn(batch_size, feature_dim)
return [x, y]

def get_init_inputs():
return [] # No special initialization inputs needed